officeparser 7.2.3 → 7.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +148 -9
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +3 -2
- package/dist/generators/BaseGenerator.d.ts +11 -0
- package/dist/generators/BaseGenerator.js +29 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +24 -14
- package/dist/generators/EpubGenerator.d.ts +18 -0
- package/dist/generators/EpubGenerator.js +242 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +266 -51
- package/dist/generators/MarkdownGenerator.d.ts +16 -0
- package/dist/generators/MarkdownGenerator.js +173 -24
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +12 -15
- package/dist/generators/TextGenerator.js +11 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +143 -6
- package/dist/officeparser.browser.iife.js +284 -188
- package/dist/officeparser.browser.mjs +284 -188
- package/dist/officeparser.browser.slim.d.ts +143 -6
- package/dist/officeparser.browser.slim.iife.js +284 -188
- package/dist/officeparser.browser.slim.mjs +284 -188
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/HtmlParser.js +284 -20
- package/dist/parsers/MarkdownParser.js +424 -33
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +142 -6
- package/dist/types.js +2 -0
- package/dist/utils/errorUtils.js +3 -2
- package/dist/utils/sanitize.d.ts +99 -0
- package/dist/utils/sanitize.js +228 -0
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +9 -5
package/README.md
CHANGED
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
|
|
4
4
|
|
|
5
|
-
**Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML)
|
|
5
|
+
**Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML) · [`epub`](https://en.wikipedia.org/wiki/EPUB)
|
|
6
6
|
|
|
7
|
-
**Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `Plain Text` · `RAG Chunks`
|
|
7
|
+
**Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `EPUB` · `Plain Text` · `RAG Chunks`
|
|
8
8
|
|
|
9
9
|
[](https://badge.fury.io/js/officeparser)
|
|
10
10
|
[](https://www.npmjs.com/package/officeparser)
|
|
@@ -42,6 +42,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
42
42
|
- [Native RAG Chunking](#native-rag-chunking)
|
|
43
43
|
- [The AST Structure](#the-ast-structure)
|
|
44
44
|
- [Deep Dive: Document Components](#deep-dive-document-components)
|
|
45
|
+
- [Markdown Dialect Support](#markdown-dialect-support)
|
|
46
|
+
- [EPUB Support](#epub-support)
|
|
45
47
|
- [Performance Highlights](#performance-highlights)
|
|
46
48
|
- [Advanced AST Usage](#advanced-ast-usage)
|
|
47
49
|
- [Configuration Reference](#configuration-reference)
|
|
@@ -93,6 +95,9 @@ npx officeparser data.xlsx --to=csv --csvDelimiter=";"
|
|
|
93
95
|
# Generate RAG chunks
|
|
94
96
|
npx officeparser document.pdf --to=chunks
|
|
95
97
|
|
|
98
|
+
# Convert DOCX to EPUB (--extractAttachments is required to embed images)
|
|
99
|
+
npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
|
|
100
|
+
|
|
96
101
|
# Overriding file extension mapping
|
|
97
102
|
npx officeparser my_document --fileType=docx --to=json
|
|
98
103
|
```
|
|
@@ -106,9 +111,9 @@ npx officeparser my_document --fileType=docx --to=json
|
|
|
106
111
|
|
|
107
112
|
| Flag | Values | Default | Description |
|
|
108
113
|
|------|--------|---------|-------------|
|
|
109
|
-
| `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
|
|
114
|
+
| `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | Output format |
|
|
110
115
|
| `--output` | path | — | Write output to a file |
|
|
111
|
-
| `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
|
|
116
|
+
| `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html\|epub` | — | Explicitly override input file type detection |
|
|
112
117
|
| `--ocr` | boolean | `false` | Enable OCR for images |
|
|
113
118
|
| `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
|
|
114
119
|
| `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
|
|
@@ -126,7 +131,7 @@ npx officeparser my_document --fileType=docx --to=json
|
|
|
126
131
|
| `--includeFormatting` | boolean | `true` | Include formatting style map matching |
|
|
127
132
|
| `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
|
|
128
133
|
| `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
|
|
129
|
-
| ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
|
|
134
|
+
| ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | **Deprecated.** Use `--to` |
|
|
130
135
|
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text` |
|
|
131
136
|
| ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
|
|
132
137
|
| ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
|
|
@@ -306,13 +311,16 @@ const { value: html } = await OfficeGenerator.generate(ast, 'html', {
|
|
|
306
311
|
const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
|
|
307
312
|
```
|
|
308
313
|
|
|
309
|
-
**Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'chunks'`
|
|
314
|
+
**Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'epub'` · `'chunks'`
|
|
310
315
|
|
|
311
316
|
> [!NOTE]
|
|
312
317
|
> **PDF generation** requires the optional `puppeteer` peer dependency:
|
|
313
318
|
> ```bash
|
|
314
319
|
> npm install puppeteer
|
|
315
320
|
> ```
|
|
321
|
+
>
|
|
322
|
+
> **EPUB generation with images** requires `extractAttachments: true` on the parse step that
|
|
323
|
+
> produced the AST — see [EPUB Support](#epub-support).
|
|
316
324
|
|
|
317
325
|
---
|
|
318
326
|
|
|
@@ -440,10 +448,10 @@ interface OfficeChunk {
|
|
|
440
448
|
|
|
441
449
|
```text
|
|
442
450
|
OfficeParserAST
|
|
443
|
-
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (
|
|
451
|
+
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | 'epub' | ... (12 formats)
|
|
444
452
|
├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
|
|
445
453
|
├── content: [ OfficeContentNode ]
|
|
446
|
-
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
|
|
454
|
+
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | 'admonition' | 'embed' | 'definitionList' | ...
|
|
447
455
|
│ ├── text: string (concatenated text of node + all descendants)
|
|
448
456
|
│ ├── children: [ OfficeContentNode ] (recursive structural children)
|
|
449
457
|
│ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
|
|
@@ -596,6 +604,94 @@ console.log(ast.metadata.nativeProperties);
|
|
|
596
604
|
// PDF: { Title: 'Report', XMP: { ... } }
|
|
597
605
|
```
|
|
598
606
|
|
|
607
|
+
### 8. Admonitions, Embeds & Definition Lists
|
|
608
|
+
|
|
609
|
+
```text
|
|
610
|
+
Admonition Node (type: 'admonition')
|
|
611
|
+
├── metadata: { admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution', title?: string }
|
|
612
|
+
└── children: [ Paragraph | List | ... ] (block content)
|
|
613
|
+
|
|
614
|
+
Embed Node (type: 'embed')
|
|
615
|
+
└── metadata: { embedType: 'youtube', videoId: string, url?: string, width?: string, align?: string }
|
|
616
|
+
|
|
617
|
+
Definition List Node (type: 'definitionList')
|
|
618
|
+
└── children:
|
|
619
|
+
├── Definition Term (type: 'definitionTerm')
|
|
620
|
+
└── Definition Description (type: 'definitionDescription')
|
|
621
|
+
```
|
|
622
|
+
|
|
623
|
+
- `admonition` round-trips through both Markdown (`> [!NOTE]` / `:::note ... :::`) and HTML (`<div class="admonition admonition-note" data-type="note">`)
|
|
624
|
+
- `embed` currently models YouTube videos; HTML round-trips via `<div data-youtube-video="ID">`, Markdown falls back to a raw HTML block or a plain link
|
|
625
|
+
- Abbreviations (`*[HTML]: Hypertext Markup Language`) are stored as `TextMetadata.abbreviationTitle` on the abbreviated text node rather than as a separate node type
|
|
626
|
+
|
|
627
|
+
---
|
|
628
|
+
|
|
629
|
+
## Markdown Dialect Support
|
|
630
|
+
|
|
631
|
+
Beyond CommonMark/GFM basics, `MarkdownParser`/`MarkdownGenerator` support an extended dialect aimed at
|
|
632
|
+
full-fidelity round-tripping with rich Markdown editors. Every construct below parses to a first-class
|
|
633
|
+
AST node/metadata field and regenerates back to the canonical syntax shown, so `.md → AST → .md` is
|
|
634
|
+
idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
|
|
635
|
+
|
|
636
|
+
| Feature | Markdown syntax | AST representation |
|
|
637
|
+
|---|---|---|
|
|
638
|
+
| Task lists (GFM) | `- [x] Done` / `- [ ] Todo` | `ListMetadata.isTask` / `.checked` |
|
|
639
|
+
| Admonitions | `> [!NOTE]` (also accepts GLFM `:::note ... :::` on import) | `type: 'admonition'`, `AdmonitionMetadata` |
|
|
640
|
+
| Footnotes | `Text[^1]` + `[^1]: Definition` | `type: 'note'`, keyed by footnote id |
|
|
641
|
+
| Definition lists | `Term\n: Definition` | `type: 'definitionList'` / `'definitionTerm'` / `'definitionDescription'` |
|
|
642
|
+
| Abbreviations | `*[HTML]: Hypertext Markup Language` | `TextMetadata.abbreviationTitle` |
|
|
643
|
+
| Attribute lists | `{width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
|
|
644
|
+
| Citations | `[@smith2024]` | `TextMetadata.citationKey` |
|
|
645
|
+
| Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
|
|
646
|
+
| Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `TextMetadata.math` (`'inline' \| 'block'`) |
|
|
647
|
+
| Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
|
|
648
|
+
| MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
|
|
649
|
+
|
|
650
|
+
> [!NOTE]
|
|
651
|
+
> MDX/JSX stripping is one-directional (parse-only) — officeParser never authors JSX back into Markdown.
|
|
652
|
+
> Wikilink enable/disable and citekey→bibliography resolution are application-level concerns; officeParser
|
|
653
|
+
> always parses/generates the syntax itself.
|
|
654
|
+
|
|
655
|
+
The same round-trip fidelity extends to HTML, so content saved from a rich-text editor survives a
|
|
656
|
+
save→reload cycle:
|
|
657
|
+
|
|
658
|
+
| HTML attribute | AST field | Notes |
|
|
659
|
+
|---|---|---|
|
|
660
|
+
| `data-width` / `data-align` / inline `style="width:…"` on `<img>` | `ImageMetadata.width` / `.align` | |
|
|
661
|
+
| `data-align` on `<table>` | `TableMetadata.align` | |
|
|
662
|
+
| `colspan` / `rowspan` on `<td>`/`<th>` | `CellMetadata.colSpan` / `.rowSpan` | Previously dropped on HTML import — merged cells now survive a save→reload cycle |
|
|
663
|
+
| `<div data-youtube-video="ID">` / `<iframe src="...youtube.com...">` | `type: 'embed'` | |
|
|
664
|
+
| `<ul data-type="taskList">` / `<li data-checked>` | `ListMetadata.isTask` / `.checked` | |
|
|
665
|
+
|
|
666
|
+
---
|
|
667
|
+
|
|
668
|
+
## EPUB Support
|
|
669
|
+
|
|
670
|
+
EPUB files are ZIP archives of XHTML content plus an OPF manifest — `EpubParser` unzips the archive,
|
|
671
|
+
resolves the spine's reading order from `content.opf`, and parses each XHTML document through the
|
|
672
|
+
existing `HtmlParser`, so EPUB content shares the same AST shape (and the same Markdown-dialect
|
|
673
|
+
fidelity above) as every other format. Dublin Core metadata (`dc:title`, `dc:creator`, `dc:description`,
|
|
674
|
+
`dc:subject`, `dc:date`, `dc:publisher`, `dc:language`, `dc:identifier`) maps into `ast.metadata` /
|
|
675
|
+
`ast.metadata.nativeProperties`, and cover art is exposed via `metadata.customProperties.coverImageName`.
|
|
676
|
+
|
|
677
|
+
`EpubGenerator` renders the AST through `HtmlGenerator` and packages the result as a minimal, valid
|
|
678
|
+
EPUB 3 (`mimetype`, `META-INF/container.xml`, an OPF manifest, a nav document, and one XHTML chapter).
|
|
679
|
+
|
|
680
|
+
> [!IMPORTANT]
|
|
681
|
+
> **Pass `extractAttachments: true` when converting to or from EPUB if the document has images.**
|
|
682
|
+
> Without it, the parser never pulls embedded image bytes out of the source document, so there is
|
|
683
|
+
> nothing for the EPUB generator to package — images silently disappear even though everything else
|
|
684
|
+
> converts correctly. Images are packaged as real zip entries (`OEBPS/images/...`) declared in the OPF
|
|
685
|
+
> manifest, not `data:` URIs — most EPUB reading systems do not render `data:` URIs in image `src`.
|
|
686
|
+
>
|
|
687
|
+
> This only matters for the two-step `OfficeParser.parseOffice()` → `OfficeGenerator.generate()` API
|
|
688
|
+
> and the CLI. [`OfficeConverter.convert()`](#officeconverter-one-step-api) enables `extractAttachments`
|
|
689
|
+
> automatically unless you explicitly set `generatorConfig.includeImages: false`.
|
|
690
|
+
>
|
|
691
|
+
> ```bash
|
|
692
|
+
> npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
|
|
693
|
+
> ```
|
|
694
|
+
|
|
599
695
|
---
|
|
600
696
|
|
|
601
697
|
## Performance Highlights
|
|
@@ -852,7 +948,7 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
852
948
|
|
|
853
949
|
| Option | Type | Default | Description |
|
|
854
950
|
|--------|------|---------|-------------|
|
|
855
|
-
| `standalone` | `boolean` | `true` |
|
|
951
|
+
| `standalone` | `boolean \| StandaloneConfig` | `true` | Controls the HTML "document envelope" — see below |
|
|
856
952
|
| `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
|
|
857
953
|
| `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
|
|
858
954
|
| `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
|
|
@@ -861,6 +957,49 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
861
957
|
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
862
958
|
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
863
959
|
|
|
960
|
+
#### `standalone`: granular envelope control
|
|
961
|
+
|
|
962
|
+
`standalone` conflates several independent decisions: whether to emit the `<!doctype>/<html>/<head>/
|
|
963
|
+
<body>` shell, how CSS is delivered, and whether to inject scripts/meta tags/injections. The boolean
|
|
964
|
+
shorthand still works — **`true`/omitted turns every part on** (a complete document); **`false` turns
|
|
965
|
+
every part off** (a bare content fragment, safe to drop into a page you don't control). Pass an
|
|
966
|
+
object instead for granular control; any field you omit defaults to its "on" (standalone) value:
|
|
967
|
+
|
|
968
|
+
| `StandaloneConfig` field | Type | Default | Description |
|
|
969
|
+
|--------|------|---------|-------------|
|
|
970
|
+
| `document` | `boolean` | `true` | Wrap in `<!DOCTYPE html><html><head>…</head><body>…</body></html>` |
|
|
971
|
+
| `metaTags` | `boolean` | `true` | Emit `<title>`/`<meta>` tags. Only meaningful when `document` is true |
|
|
972
|
+
| `styles` | `'full' \| 'scoped' \| 'none'` | `'full'` | See below |
|
|
973
|
+
| `scripts` | `boolean` | `true` | Emit the Chart.js CDN loader and spreadsheet-interactivity `<script>` tags |
|
|
974
|
+
| `headInjections` | `boolean` | `true` | Apply `injections.headStart`/`headEnd`. Only meaningful when `document` is true |
|
|
975
|
+
| `bodyInjections` | `boolean` | `true` | Apply `injections.bodyStart`/`bodyEnd` — applies even to a bare fragment |
|
|
976
|
+
|
|
977
|
+
`styles` controls how the built-in stylesheet is delivered:
|
|
978
|
+
- **`'full'`** — the complete stylesheet using global selectors (`body`, `h1`, `table`, …). This is
|
|
979
|
+
what `standalone: true` has always emitted.
|
|
980
|
+
- **`'scoped'`** — the same styling, scoped under the fragment's own wrapper via CSS `@scope` so it
|
|
981
|
+
cannot leak onto a host page's elements. Requires a modern engine (Chrome 118+, Safari 17.4+,
|
|
982
|
+
Firefox 128+); for universal support use `'none'` (bring your own CSS) or `'full'`.
|
|
983
|
+
- **`'none'`** — no stylesheet at all; the host page (or rich-text editor, or EPUB reader) supplies
|
|
984
|
+
its own styling.
|
|
985
|
+
|
|
986
|
+
```js
|
|
987
|
+
// A styled fragment to embed in your own page, without a document shell:
|
|
988
|
+
await ast.to('html', { htmlConfig: { standalone: { document: false } } });
|
|
989
|
+
|
|
990
|
+
// The same, but with styles scoped so they can't leak onto your page's own elements:
|
|
991
|
+
await ast.to('html', { htmlConfig: { standalone: { document: false, styles: 'scoped' } } });
|
|
992
|
+
|
|
993
|
+
// A completely bare fragment (no shell, no styles, no scripts) — e.g. for a rich-text editor:
|
|
994
|
+
await ast.to('html', { htmlConfig: { standalone: false } });
|
|
995
|
+
```
|
|
996
|
+
|
|
997
|
+
> [!NOTE]
|
|
998
|
+
> **Behavior change from `standalone: false`:** previously this emitted a fragment with a *global,
|
|
999
|
+
> unscoped* `<style>` block. It now emits a genuinely bare fragment (no `<style>` at all), matching
|
|
1000
|
+
> "every part off." If you relied on the old styled-fragment behavior, pass
|
|
1001
|
+
> `{ document: false }` (or `{ document: false, styles: 'full' }`) instead.
|
|
1002
|
+
|
|
864
1003
|
### MdGeneratorConfig
|
|
865
1004
|
|
|
866
1005
|
Pass as `mdConfig` inside `GeneratorConfig`.
|
package/dist/OfficeGenerator.js
CHANGED
|
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.OfficeGenerator = void 0;
|
|
4
4
|
const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
|
|
5
5
|
const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
|
|
6
|
+
const EpubGenerator_js_1 = require("./generators/EpubGenerator.js");
|
|
6
7
|
const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
|
|
7
8
|
const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
|
|
8
9
|
const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
|
|
@@ -60,6 +61,9 @@ class OfficeGenerator {
|
|
|
60
61
|
case 'chunks':
|
|
61
62
|
generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
|
|
62
63
|
break;
|
|
64
|
+
case 'epub':
|
|
65
|
+
generator = new EpubGenerator_js_1.EpubGenerator(ast, config);
|
|
66
|
+
break;
|
|
63
67
|
default:
|
|
64
68
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
|
|
65
69
|
}
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
* - CSV (Comma-Separated Values)
|
|
15
15
|
* - MD (Markdown)
|
|
16
16
|
* - HTML (HyperText Markup Language)
|
|
17
|
+
* - EPUB (E-book format)
|
|
17
18
|
*
|
|
18
19
|
* **Usage:**
|
|
19
20
|
* ```typescript
|
|
@@ -66,6 +67,7 @@ export declare class OfficeParser {
|
|
|
66
67
|
* - `.csv` → CsvParser
|
|
67
68
|
* - `.md` → MarkdownParser
|
|
68
69
|
* - `.html` → HtmlParser
|
|
70
|
+
* - `.epub` → EpubParser
|
|
69
71
|
*
|
|
70
72
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
71
73
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
package/dist/OfficeParser.js
CHANGED
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* - CSV (Comma-Separated Values)
|
|
16
16
|
* - MD (Markdown)
|
|
17
17
|
* - HTML (HyperText Markup Language)
|
|
18
|
+
* - EPUB (E-book format)
|
|
18
19
|
*
|
|
19
20
|
* **Usage:**
|
|
20
21
|
* ```typescript
|
|
@@ -39,6 +40,7 @@
|
|
|
39
40
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
40
41
|
exports.OfficeParser = void 0;
|
|
41
42
|
const CsvParser_js_1 = require("./parsers/CsvParser.js");
|
|
43
|
+
const EpubParser_js_1 = require("./parsers/EpubParser.js");
|
|
42
44
|
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
43
45
|
const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
|
|
44
46
|
const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
|
|
@@ -83,6 +85,7 @@ class OfficeParser {
|
|
|
83
85
|
* - `.csv` → CsvParser
|
|
84
86
|
* - `.md` → MarkdownParser
|
|
85
87
|
* - `.html` → HtmlParser
|
|
88
|
+
* - `.epub` → EpubParser
|
|
86
89
|
*
|
|
87
90
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
88
91
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
@@ -232,6 +235,9 @@ class OfficeParser {
|
|
|
232
235
|
case 'md':
|
|
233
236
|
result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
|
|
234
237
|
break;
|
|
238
|
+
case 'epub':
|
|
239
|
+
result = await (0, EpubParser_js_1.parseEpub)(buffer, internalConfig);
|
|
240
|
+
break;
|
|
235
241
|
default:
|
|
236
242
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
237
243
|
}
|
package/dist/cli.d.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* officeparser file.docx --ocr --extractAttachments
|
|
9
9
|
*
|
|
10
10
|
* Options (--key=value, --key value, or bare flags):
|
|
11
|
-
* --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
|
|
11
|
+
* --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
|
|
12
12
|
* --output=path Save result to a file
|
|
13
13
|
* --fileType=docx|xlsx|... Override file type detection
|
|
14
14
|
* --ocr Enable OCR for images (default: false)
|
package/dist/cli.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* officeparser file.docx --ocr --extractAttachments
|
|
10
10
|
*
|
|
11
11
|
* Options (--key=value, --key value, or bare flags):
|
|
12
|
-
* --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
|
|
12
|
+
* --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
|
|
13
13
|
* --output=path Save result to a file
|
|
14
14
|
* --fileType=docx|xlsx|... Override file type detection
|
|
15
15
|
* --ocr Enable OCR for images (default: false)
|
|
@@ -333,7 +333,7 @@ else {
|
|
|
333
333
|
console.log('Usage: officeparser <file> [options]');
|
|
334
334
|
console.log('');
|
|
335
335
|
console.log('Options:');
|
|
336
|
-
console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks
|
|
336
|
+
console.log(' --to=json|text|md|html|pdf|csv|rtf|epub|chunks Target conversion format (default: json)');
|
|
337
337
|
console.log(' --output=file.ext Save output to file instead of stdout');
|
|
338
338
|
console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
|
|
339
339
|
console.log(' --ocr Enable OCR for images (default: false)');
|
|
@@ -371,5 +371,6 @@ else {
|
|
|
371
371
|
console.log(' officeparser document.docx --to md');
|
|
372
372
|
console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
|
|
373
373
|
console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
|
|
374
|
+
console.log(' officeparser document.docx --extractAttachments --to epub --output document.epub');
|
|
374
375
|
console.log(' officeparser image_doc --fileType docx --to json');
|
|
375
376
|
}
|
|
@@ -48,6 +48,17 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
|
|
|
48
48
|
* Helper to generate a unique ID (slug) from text.
|
|
49
49
|
*/
|
|
50
50
|
protected slugify(text: string): string;
|
|
51
|
+
private noteFootnoteKeys;
|
|
52
|
+
private usedFootnoteKeys;
|
|
53
|
+
private footnoteKeyCounter;
|
|
54
|
+
/**
|
|
55
|
+
* Assigns a stable, unique reference key to a footnote/endnote node, reused for both
|
|
56
|
+
* its inline reference marker and its collected definition. Source ids aren't reliably
|
|
57
|
+
* unique across a document - DOCX/ODT number footnotes and endnotes in separate
|
|
58
|
+
* sequences, so both can carry noteId "1" - so a source id is only reused when it
|
|
59
|
+
* hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
|
|
60
|
+
*/
|
|
61
|
+
protected getFootnoteKey(note: OfficeContentNode): string;
|
|
51
62
|
/**
|
|
52
63
|
* Recursively extracts plain text from a node and its children.
|
|
53
64
|
*/
|
|
@@ -89,6 +89,35 @@ class BaseGenerator {
|
|
|
89
89
|
.replace(/[\s_-]+/g, '-')
|
|
90
90
|
.replace(/^-+|-+$/g, '');
|
|
91
91
|
}
|
|
92
|
+
noteFootnoteKeys = new Map();
|
|
93
|
+
usedFootnoteKeys = new Set();
|
|
94
|
+
footnoteKeyCounter = 0;
|
|
95
|
+
/**
|
|
96
|
+
* Assigns a stable, unique reference key to a footnote/endnote node, reused for both
|
|
97
|
+
* its inline reference marker and its collected definition. Source ids aren't reliably
|
|
98
|
+
* unique across a document - DOCX/ODT number footnotes and endnotes in separate
|
|
99
|
+
* sequences, so both can carry noteId "1" - so a source id is only reused when it
|
|
100
|
+
* hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
|
|
101
|
+
*/
|
|
102
|
+
getFootnoteKey(note) {
|
|
103
|
+
const cached = this.noteFootnoteKeys.get(note);
|
|
104
|
+
if (cached)
|
|
105
|
+
return cached;
|
|
106
|
+
const preferred = note.metadata?.noteId;
|
|
107
|
+
let key;
|
|
108
|
+
if (preferred && !this.usedFootnoteKeys.has(preferred)) {
|
|
109
|
+
key = preferred;
|
|
110
|
+
}
|
|
111
|
+
else {
|
|
112
|
+
do {
|
|
113
|
+
this.footnoteKeyCounter++;
|
|
114
|
+
key = String(this.footnoteKeyCounter);
|
|
115
|
+
} while (this.usedFootnoteKeys.has(key));
|
|
116
|
+
}
|
|
117
|
+
this.usedFootnoteKeys.add(key);
|
|
118
|
+
this.noteFootnoteKeys.set(note, key);
|
|
119
|
+
return key;
|
|
120
|
+
}
|
|
92
121
|
/**
|
|
93
122
|
* Recursively extracts plain text from a node and its children.
|
|
94
123
|
*/
|
|
@@ -20,9 +20,17 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
|
|
|
20
20
|
*/
|
|
21
21
|
private renderNodeToRows;
|
|
22
22
|
/**
|
|
23
|
-
* Escapes a value for CSV formatting
|
|
23
|
+
* Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
|
|
24
|
+
* formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
|
|
24
25
|
*/
|
|
25
26
|
private escapeCsvValue;
|
|
27
|
+
/** Sanitizes a value for a `#` comment line (document metadata or sheet name).
|
|
28
|
+
* Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
|
|
29
|
+
* can't be quoted; instead we (a) collapse newlines so the value can't break out
|
|
30
|
+
* and inject a new row, and (b) replace the column delimiter with a space so a
|
|
31
|
+
* value like `x,=1+1` can't split into a second cell that a spreadsheet would
|
|
32
|
+
* evaluate as a formula (CSV formula/DDE injection). */
|
|
33
|
+
private sanitizeComment;
|
|
26
34
|
/**
|
|
27
35
|
* Renders metadata as comments.
|
|
28
36
|
*/
|
|
@@ -4,6 +4,7 @@ exports.CsvGenerator = void 0;
|
|
|
4
4
|
const fflate_1 = require("fflate");
|
|
5
5
|
const types_js_1 = require("../types.js");
|
|
6
6
|
const sheetUtils_js_1 = require("../utils/sheetUtils.js");
|
|
7
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
7
8
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
8
9
|
/**
|
|
9
10
|
* Generates CSV files from an AST.
|
|
@@ -26,7 +27,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
26
27
|
if (sheetNodes.length === 0) {
|
|
27
28
|
return { value: '', messages: this.messages };
|
|
28
29
|
}
|
|
29
|
-
const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast) : '';
|
|
30
|
+
const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast, delimiter) : '';
|
|
30
31
|
// 2. Filter sheets based on range
|
|
31
32
|
let selectedNodes = sheetNodes;
|
|
32
33
|
if (csvConfig.sheets) {
|
|
@@ -59,7 +60,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
59
60
|
mergedLines.push(metadataHeader.trim());
|
|
60
61
|
for (const sheet of sheetData) {
|
|
61
62
|
if (sheetData.length > 1)
|
|
62
|
-
mergedLines.push(`# Sheet: ${sheet.name}`);
|
|
63
|
+
mergedLines.push(`# Sheet: ${this.sanitizeComment(sheet.name, delimiter)}`);
|
|
63
64
|
for (const row of sheet.rows) {
|
|
64
65
|
const paddedRow = [...row];
|
|
65
66
|
// Don't pad comments
|
|
@@ -197,34 +198,43 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
197
198
|
return rows;
|
|
198
199
|
}
|
|
199
200
|
/**
|
|
200
|
-
* Escapes a value for CSV formatting
|
|
201
|
+
* Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
|
|
202
|
+
* formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
|
|
201
203
|
*/
|
|
202
204
|
escapeCsvValue(val, delimiter) {
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
205
|
+
return (0, sanitize_js_1.csvSafeCell)(val, delimiter);
|
|
206
|
+
}
|
|
207
|
+
/** Sanitizes a value for a `#` comment line (document metadata or sheet name).
|
|
208
|
+
* Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
|
|
209
|
+
* can't be quoted; instead we (a) collapse newlines so the value can't break out
|
|
210
|
+
* and inject a new row, and (b) replace the column delimiter with a space so a
|
|
211
|
+
* value like `x,=1+1` can't split into a second cell that a spreadsheet would
|
|
212
|
+
* evaluate as a formula (CSV formula/DDE injection). */
|
|
213
|
+
sanitizeComment(val, delimiter) {
|
|
214
|
+
let s = String(val ?? '').replace(/[\r\n]+/g, ' ');
|
|
215
|
+
if (delimiter)
|
|
216
|
+
s = s.split(delimiter).join(' ');
|
|
217
|
+
return s;
|
|
208
218
|
}
|
|
209
219
|
/**
|
|
210
220
|
* Renders metadata as comments.
|
|
211
221
|
*/
|
|
212
|
-
renderMetadata(ast) {
|
|
222
|
+
renderMetadata(ast, delimiter) {
|
|
213
223
|
if (!ast.metadata)
|
|
214
224
|
return '';
|
|
215
225
|
const m = ast.metadata;
|
|
216
226
|
let output = '';
|
|
217
227
|
if (m.title)
|
|
218
|
-
output += `# Title: ${m.title}\n`;
|
|
228
|
+
output += `# Title: ${this.sanitizeComment(m.title, delimiter)}\n`;
|
|
219
229
|
if (m.author)
|
|
220
|
-
output += `# Author: ${m.author}\n`;
|
|
230
|
+
output += `# Author: ${this.sanitizeComment(m.author, delimiter)}\n`;
|
|
221
231
|
if (m.created)
|
|
222
|
-
output += `# Created: ${new Date(m.created).toLocaleString()}\n`;
|
|
232
|
+
output += `# Created: ${this.sanitizeComment(new Date(m.created).toLocaleString(), delimiter)}\n`;
|
|
223
233
|
if (m.modified)
|
|
224
|
-
output += `# Modified: ${new Date(m.modified).toLocaleString()}\n`;
|
|
234
|
+
output += `# Modified: ${this.sanitizeComment(new Date(m.modified).toLocaleString(), delimiter)}\n`;
|
|
225
235
|
if (m.customProperties) {
|
|
226
236
|
for (const [k, v] of Object.entries(m.customProperties)) {
|
|
227
|
-
output += `# ${k}: ${v}\n`;
|
|
237
|
+
output += `# ${this.sanitizeComment(k, delimiter)}: ${this.sanitizeComment(String(v), delimiter)}\n`;
|
|
228
238
|
}
|
|
229
239
|
}
|
|
230
240
|
return output ? output + '\n' : '';
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST } from '../types.js';
|
|
2
|
+
import { BaseGenerator } from './BaseGenerator.js';
|
|
3
|
+
/**
|
|
4
|
+
* Generates a minimal, valid EPUB 3 file from an AST.
|
|
5
|
+
*
|
|
6
|
+
* Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
|
|
7
|
+
* for the actual markup, since EPUB content documents are XHTML) and packaged with the
|
|
8
|
+
* required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
|
|
9
|
+
*
|
|
10
|
+
* `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
|
|
11
|
+
* render `data:` URIs - images must be packaged as separate resources referenced by a
|
|
12
|
+
* relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
|
|
13
|
+
* the manifest, and its `<img src>` rewritten to point at the packaged file.
|
|
14
|
+
*/
|
|
15
|
+
export declare class EpubGenerator extends BaseGenerator<'epub'> {
|
|
16
|
+
constructor(ast: OfficeParserAST, config?: GeneratorConfig<'epub'>);
|
|
17
|
+
generate(): Promise<ConversionResult<'epub'>>;
|
|
18
|
+
}
|